
  <rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
    <channel>
      <title>HANS</title>
      <link>https://www.hansking.cn/blog</link>
      <description>A posts created with Next.js and Tailwind.css</description>
      <language>en-us</language>
      <managingEditor>admin@hansking.cn (HANS)</managingEditor>
      <webMaster>admin@hansking.cn (HANS)</webMaster>
      <lastBuildDate>Sun, 26 Oct 2025 10:00:00 GMT</lastBuildDate>
      <atom:link href="https://www.hansking.cn/tags/Speculative Decoding/feed.xml" rel="self" type="application/rss+xml"/>
      
  <item>
    <guid>https://www.hansking.cn/blog/2026/SpeculativeDecoding-yong-xiao-mo-xing-jia-su-da-mo-xing</guid>
    <title>Speculative Decoding：用小模型加速大模型</title>
    <link>https://www.hansking.cn/blog/2026/SpeculativeDecoding-yong-xiao-mo-xing-jia-su-da-mo-xing</link>
    <description>[{&quot;value&quot;:&quot;Speculative Decoding：用小模型加速大模型&quot;,&quot;url&quot;:&quot;#speculative-decoding用小模型加速大模型&quot;,&quot;depth&quot;:1},{&quot;value&quot;:&quot;一、为什么自回归生成慢&quot;,&quot;url&quot;:&quot;#一为什么自回归生成慢&quot;,&quot;depth&quot;:2},{&quot;value&quot;:&quot;二、基本流程是什么&quot;,&quot;url&quot;:&quot;#二基本流程是什么&quot;,&quot;depth&quot;:2},{&quot;value&quot;:&quot;三、小模型不需要和大模型一样聪明&quot;,&quot;url&quot;:&quot;#三小模型不需要和大模型一样聪明&quot;,&quot;depth&quot;:2},{&quot;value&quot;:&quot;四、速度收益取决于接受率&quot;,&quot;url&quot;:&quot;#四速度收益取决于接受率&quot;,&quot;depth&quot;:2},{&quot;value&quot;:&quot;五、为什么结果可以保持目标模型分布&quot;,&quot;url&quot;:&quot;#五为什么结果可以保持目标模型分布&quot;,&quot;depth&quot;:2},{&quot;value&quot;:&quot;六、KV Cache 管理决定能不能跑得快&quot;,&quot;url&quot;:&quot;#六kv-cache-管理决定能不能跑得快&quot;,&quot;depth&quot;:2},{&quot;value&quot;:&quot;七、不同任务的收益不同&quot;,&quot;url&quot;:&quot;#七不同任务的收益不同&quot;,&quot;depth&quot;:2},{&quot;value&quot;:&quot;八、工程上常见的失败原因&quot;,&quot;url&quot;:&quot;#八工程上常见的失败原因&quot;,&quot;depth&quot;:2},{&quot;value&quot;:&quot;1. 草稿模型太慢&quot;,&quot;url&quot;:&quot;#1-草稿模型太慢&quot;,&quot;depth&quot;:3},{&quot;value&quot;:&quot;2. Tokenizer 不兼容&quot;,&quot;url&quot;:&quot;#2-tokenizer-不兼容&quot;,&quot;depth&quot;:3},{&quot;value&quot;:&quot;3. 采样参数不一致&quot;,&quot;url&quot;:&quot;#3-采样参数不一致&quot;,&quot;depth&quot;:3},{&quot;value&quot;:&quot;4. 并发下显存不足&quot;,&quot;url&quot;:&quot;#4-并发下显存不足&quot;,&quot;depth&quot;:3},{&quot;value&quot;:&quot;5. 只测平均值&quot;,&quot;url&quot;:&quot;#5-只测平均值&quot;,&quot;depth&quot;:3},{&quot;value&quot;:&quot;九、上线前怎么评估&quot;,&quot;url&quot;:&quot;#九上线前怎么评估&quot;,&quot;depth&quot;:2},{&quot;value&quot;:&quot;总结：让小模型猜，让大模型负责&quot;,&quot;url&quot;:&quot;#总结让小模型猜让大模型负责&quot;,&quot;depth&quot;:2}]</description>
    <pubDate>Sun, 26 Oct 2025 10:00:00 GMT</pubDate>
    <author>admin@hansking.cn (HANS)</author>
    <category>AI</category><category>LLM</category><category>Speculative Decoding</category><category>推理优化</category><category>性能优化</category>
  </item>

    </channel>
  </rss>
